agora inbox for [email protected]help / color / mirror / Atom feed
[PATCH v1 4/7] Row pattern recognition patch (executor). 189+ messages / 2 participants [nested] [flat]
* [PATCH v1 4/7] Row pattern recognition patch (executor). @ 2023-06-25 11:48 Tatsuo Ishii <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Tatsuo Ishii @ 2023-06-25 11:48 UTC (permalink / raw) --- src/backend/executor/nodeWindowAgg.c | 223 +++++++++++++++++++++- src/backend/utils/adt/windowfuncs.c | 274 ++++++++++++++++++++++++++- src/include/catalog/pg_proc.dat | 9 + src/include/nodes/execnodes.h | 12 ++ src/include/windowapi.h | 9 + 5 files changed, 517 insertions(+), 10 deletions(-) diff --git a/src/backend/executor/nodeWindowAgg.c b/src/backend/executor/nodeWindowAgg.c index 310ac23e3a..ae997dff86 100644 --- a/src/backend/executor/nodeWindowAgg.c +++ b/src/backend/executor/nodeWindowAgg.c @@ -48,6 +48,7 @@ #include "utils/acl.h" #include "utils/builtins.h" #include "utils/datum.h" +#include "utils/fmgroids.h" #include "utils/expandeddatum.h" #include "utils/lsyscache.h" #include "utils/memutils.h" @@ -159,6 +160,14 @@ typedef struct WindowStatePerAggData bool restart; /* need to restart this agg in this cycle? */ } WindowStatePerAggData; +/* + * Map between Var attno in a target list and the parsed attno. + */ +typedef struct AttnoMap { + List *attno; /* att number in target list (list of AttNumber) */ + List *attnosyn; /* parsed att number (list of AttNumber) */ +} AttnoMap; + static void initialize_windowaggregate(WindowAggState *winstate, WindowStatePerFunc perfuncstate, WindowStatePerAgg peraggstate); @@ -195,9 +204,9 @@ static Datum GetAggInitVal(Datum textInitVal, Oid transtype); static bool are_peers(WindowAggState *winstate, TupleTableSlot *slot1, TupleTableSlot *slot2); -static bool window_gettupleslot(WindowObject winobj, int64 pos, - TupleTableSlot *slot); +static void attno_map(Node *node, AttnoMap *map); +static bool attno_map_walker(Node *node, void *context); /* * initialize_windowaggregate @@ -2388,6 +2397,12 @@ ExecInitWindowAgg(WindowAgg *node, EState *estate, int eflags) TupleDesc scanDesc; ListCell *l; + TargetEntry *te; + Expr *expr; + Var *var; + int nargs; + AttnoMap attnomap; + /* check for unsupported flags */ Assert(!(eflags & (EXEC_FLAG_BACKWARD | EXEC_FLAG_MARK))); @@ -2483,6 +2498,16 @@ ExecInitWindowAgg(WindowAgg *node, EState *estate, int eflags) winstate->temp_slot_2 = ExecInitExtraTupleSlot(estate, scanDesc, &TTSOpsMinimalTuple); + winstate->prev_slot = ExecInitExtraTupleSlot(estate, scanDesc, + &TTSOpsMinimalTuple); + + winstate->next_slot = ExecInitExtraTupleSlot(estate, scanDesc, + &TTSOpsMinimalTuple); + + winstate->null_slot = ExecInitExtraTupleSlot(estate, scanDesc, + &TTSOpsMinimalTuple); + winstate->null_slot = ExecStoreAllNullTuple(winstate->null_slot); + /* * create frame head and tail slots only if needed (must create slots in * exactly the same cases that update_frameheadpos and update_frametailpos @@ -2667,6 +2692,69 @@ ExecInitWindowAgg(WindowAgg *node, EState *estate, int eflags) winstate->inRangeAsc = node->inRangeAsc; winstate->inRangeNullsFirst = node->inRangeNullsFirst; + /* Set up row pattern recognition PATTERN clause */ + winstate->patternVariableList = node->patternVariable; + winstate->patternRegexpList = node->patternRegexp; + + /* Set up row pattern recognition DEFINE clause */ + + /* + * Collect mapping between varattno and varattnosyn in the targetlist. + * XXX: For now we only check RPR's argument. Eventually we have to + * recurse the targetlist to find out all mappings in Var nodes. + */ + attnomap.attno = NIL; + attnomap.attnosyn = NIL; + + foreach (l, node->plan.targetlist) + { + te = lfirst(l); + if (IsA(te->expr, WindowFunc)) + { + WindowFunc *func = (WindowFunc *)te->expr; + if (func->winfnoid != F_RPR) + continue; + + /* sanity check */ + nargs = list_length(func->args); + if (list_length(func->args) != 1) + elog(ERROR, "RPR must have 1 argument but function %d has %d args", func->winfnoid, nargs); + + expr = (Expr *) lfirst(list_head(func->args)); + if (!IsA(expr, Var)) + elog(ERROR, "RPR's arg is not Var"); + + var = (Var *)expr; + elog(DEBUG1, "resname: %s varattno: %d varattnosyn: %d", + te->resname, var->varattno, var->varattnosyn); + attnomap.attno = lappend_int(attnomap.attno, var->varattno); + attnomap.attnosyn = lappend_int(attnomap.attnosyn, var->varattnosyn); + } + } + + winstate->defineVariableList = NIL; + winstate->defineClauseList = NIL; + if (node->defineClause != NIL) + { + foreach(l, node->defineClause) + { + char *name; + ExprState *exps; + + te = lfirst(l); + name = te->resname; + expr = te->expr; + + elog(DEBUG1, "defineVariable name: %s", name); + winstate->defineVariableList = lappend(winstate->defineVariableList, + makeString(pstrdup(name))); + /* tweak expr so that it referes to outer slot */ + attno_map((Node *)expr, &attnomap); + exps = ExecInitExpr(expr, (PlanState *) winstate); + winstate->defineClauseList = lappend(winstate->defineClauseList, exps); + } + } + winstate->all_first = true; winstate->partition_spooled = false; winstate->more_partitions = false; @@ -2674,6 +2762,76 @@ ExecInitWindowAgg(WindowAgg *node, EState *estate, int eflags) return winstate; } +/* + * Rewrite Var node's varattno to the varattno which is used in the target + * list using AttnoMap. We also rewrite varno so that it sees outer tuple + * (PREV) or inner tuple (NEXT). + */ +static void +attno_map(Node *node, AttnoMap *map) +{ + (void) expression_tree_walker(node, attno_map_walker, (void *) map); +} + +static bool +attno_map_walker(Node *node, void *context) +{ + FuncExpr *func; + int nargs; + Expr *expr; + Var *var; + AttnoMap *attnomap; + ListCell *lc1, *lc2; + + if (node == NULL) + return false; + + attnomap = (AttnoMap *) context; + + if (IsA(node, FuncExpr)) + { + func = (FuncExpr *)node; + + if (func->funcid == F_PREV || func->funcid == F_NEXT) + { + /* sanity check */ + nargs = list_length(func->args); + if (list_length(func->args) != 1) + elog(ERROR, "PREV/NEXT must have 1 argument but function %d has %d args", func->funcid, nargs); + + expr = (Expr *) lfirst(list_head(func->args)); + if (!IsA(expr, Var)) + elog(ERROR, "PREV/NEXT's arg is not Var"); /* XXX: is it possible that arg type is Const? */ + var = (Var *)expr; + + if (func->funcid == F_PREV) + var->varno = OUTER_VAR; + else + var->varno = INNER_VAR; + } + return expression_tree_walker(node, attno_map_walker, (void *) context); + } + else if (IsA(node, Var)) + { + var = (Var *)node; + + elog(DEBUG1, "original varno: %d varattno: %d", var->varno, var->varattno); + + forboth(lc1, attnomap->attno, lc2, attnomap->attnosyn) + { + int attno = lfirst_int(lc1); + int attnosyn = lfirst_int(lc2); + + if (var->varattno == attnosyn) + { + elog(DEBUG1, "loc: %d rewrite varattno from: %d to %d", var->location, attnosyn, attno); + var->varattno = attno; + } + } + } + return expression_tree_walker(node, attno_map_walker, (void *) context); +} + /* ----------------- * ExecEndWindowAgg * ----------------- @@ -2691,6 +2849,8 @@ ExecEndWindowAgg(WindowAggState *node) ExecClearTuple(node->agg_row_slot); ExecClearTuple(node->temp_slot_1); ExecClearTuple(node->temp_slot_2); + ExecClearTuple(node->prev_slot); + ExecClearTuple(node->next_slot); if (node->framehead_slot) ExecClearTuple(node->framehead_slot); if (node->frametail_slot) @@ -2740,6 +2900,8 @@ ExecReScanWindowAgg(WindowAggState *node) ExecClearTuple(node->agg_row_slot); ExecClearTuple(node->temp_slot_1); ExecClearTuple(node->temp_slot_2); + ExecClearTuple(node->prev_slot); + ExecClearTuple(node->next_slot); if (node->framehead_slot) ExecClearTuple(node->framehead_slot); if (node->frametail_slot) @@ -3080,7 +3242,7 @@ are_peers(WindowAggState *winstate, TupleTableSlot *slot1, * * Returns true if successful, false if no such row */ -static bool +bool window_gettupleslot(WindowObject winobj, int64 pos, TupleTableSlot *slot) { WindowAggState *winstate = winobj->winstate; @@ -3420,14 +3582,53 @@ WinGetFuncArgInFrame(WindowObject winobj, int argno, WindowAggState *winstate; ExprContext *econtext; TupleTableSlot *slot; - int64 abs_pos; - int64 mark_pos; Assert(WindowObjectIsValid(winobj)); winstate = winobj->winstate; econtext = winstate->ss.ps.ps_ExprContext; slot = winstate->temp_slot_1; + if (WinGetSlotInFrame(winobj, slot, + relpos, seektype, set_mark, + isnull, isout) == 0) + { + econtext->ecxt_outertuple = slot; + return ExecEvalExpr((ExprState *) list_nth(winobj->argstates, argno), + econtext, isnull); + } + + if (isout) + *isout = true; + *isnull = true; + return (Datum) 0; +} + +/* + * WinGetSlotInFrame + * slot: TupleTableSlot to store the result + * relpos: signed rowcount offset from the seek position + * seektype: WINDOW_SEEK_HEAD or WINDOW_SEEK_TAIL + * set_mark: If the row is found/in frame and set_mark is true, the mark is + * moved to the row as a side-effect. + * isnull: output argument, receives isnull status of result + * isout: output argument, set to indicate whether target row position + * is out of frame (can pass NULL if caller doesn't care about this) + * + * Returns 0 if we successfullt got the slot. false if out of frame. + * (also isout is set) + */ +int +WinGetSlotInFrame(WindowObject winobj, TupleTableSlot *slot, + int relpos, int seektype, bool set_mark, + bool *isnull, bool *isout) +{ + WindowAggState *winstate; + int64 abs_pos; + int64 mark_pos; + + Assert(WindowObjectIsValid(winobj)); + winstate = winobj->winstate; + switch (seektype) { case WINDOW_SEEK_CURRENT: @@ -3583,15 +3784,13 @@ WinGetFuncArgInFrame(WindowObject winobj, int argno, *isout = false; if (set_mark) WinSetMarkPosition(winobj, mark_pos); - econtext->ecxt_outertuple = slot; - return ExecEvalExpr((ExprState *) list_nth(winobj->argstates, argno), - econtext, isnull); + return 0; out_of_frame: if (isout) *isout = true; *isnull = true; - return (Datum) 0; + return -1; } /* @@ -3622,3 +3821,9 @@ WinGetFuncArgCurrent(WindowObject winobj, int argno, bool *isnull) return ExecEvalExpr((ExprState *) list_nth(winobj->argstates, argno), econtext, isnull); } + +WindowAggState * +WinGetAggState(WindowObject winobj) +{ + return winobj->winstate; +} diff --git a/src/backend/utils/adt/windowfuncs.c b/src/backend/utils/adt/windowfuncs.c index b87a624fb2..63b225271c 100644 --- a/src/backend/utils/adt/windowfuncs.c +++ b/src/backend/utils/adt/windowfuncs.c @@ -13,6 +13,9 @@ */ #include "postgres.h" +#include "catalog/pg_collation_d.h" +#include "executor/executor.h" +#include "nodes/execnodes.h" #include "nodes/supportnodes.h" #include "utils/builtins.h" #include "windowapi.h" @@ -39,7 +42,9 @@ typedef struct static bool rank_up(WindowObject winobj); static Datum leadlag_common(FunctionCallInfo fcinfo, bool forward, bool withoffset, bool withdefault); - +static bool get_slots(WindowObject winobj, WindowAggState *winstate, int current_pos); +static int evaluate_pattern(WindowObject winobj, WindowAggState *winstate, + int relpos, char *vname, char *quantifier, bool *result); /* * utility routine for *_rank functions. @@ -713,3 +718,270 @@ window_nth_value(PG_FUNCTION_ARGS) PG_RETURN_DATUM(result); } + +/* + * rpr + * allow to use "Row pattern recognition: WINDOW clause" (SQL:2016 R020) in + * the target list. + * Usage: SELECT rpr(colname) OVER (..) + * where colname is defined in PATTERN clause. + */ +Datum +window_rpr(PG_FUNCTION_ARGS) +{ +#define MAX_PATTERNS 16 /* max variables in PATTERN clause */ + + WindowObject winobj = PG_WINDOW_OBJECT(); + WindowAggState *winstate = WinGetAggState(winobj); + Datum result; + bool expression_result; + bool isnull; + int relpos; + int64 curr_pos, markpos; + ListCell *lc, *lc1; + + curr_pos = WinGetCurrentPosition(winobj); + elog(DEBUG1, "rpr is called. row: " INT64_FORMAT, curr_pos); + + /* + * Evaluate PATTERN until one of expressions is not true or out of frame. + */ + relpos = 0; + + forboth(lc, winstate->patternVariableList, lc1, winstate->patternRegexpList) + { + char *vname = strVal(lfirst(lc)); + char *quantifier = strVal(lfirst(lc1)); + + elog(DEBUG1, "relpos: %d pattern vname: %s quantifier: %s", relpos, vname, quantifier); + + /* evaluate row pattern against current row */ + relpos = evaluate_pattern(winobj, winstate, relpos, vname, quantifier, &expression_result); + + /* + * If the expression did not match, we are done. + */ + if (!expression_result) + break; + + /* out of frame? */ + if (relpos < 0) + break; + + /* count up relative row position */ + relpos++; + } + + elog(DEBUG1, "relpos: %d", relpos); + + /* + * If current row satified the pattern, return argument expression. + */ + if (expression_result) + { + result = WinGetFuncArgInFrame(winobj, 0, + 0, WINDOW_SEEK_HEAD, false, + &isnull, NULL); + } + + /* + * At this point we can set mark down to current pos -2. + */ + markpos = curr_pos -2; + elog(DEBUG1, "markpos: " INT64_FORMAT, markpos); + if (markpos >= 0) + WinSetMarkPosition(winobj, markpos); + + if (expression_result) + PG_RETURN_DATUM(result); + + PG_RETURN_NULL(); +} + +/* + * Evaluate expression associated with PATTERN variable vname. + * relpos is relative row position in a frame (starting from 0). + * "quantifier" is the quatifier part of the PATTERN regular expression. + * Currently only '+' is allowed. + * result is out paramater representing the expression evaluation result + * is true of false. + * Return values are: + * >=0: the last match relative row position + * -1: current row is out of frame + */ +static +int evaluate_pattern(WindowObject winobj, WindowAggState *winstate, + int relpos, char *vname, char *quantifier, bool *result) +{ + ExprContext *econtext = winstate->ss.ps.ps_ExprContext; + ListCell *lc1, *lc2; + ExprState *pat; + Datum eval_result; + int sts; + bool out_of_frame = false; + bool isnull; + StringInfo encoded_str = makeStringInfo(); + char pattern_str[128]; + + forboth (lc1, winstate->defineVariableList, lc2, winstate->defineClauseList) + { + char *name = strVal(lfirst(lc1)); + bool second_try_match = false; + + if (strcmp(vname, name)) + continue; + + /* set expression to evaluate */ + pat = lfirst(lc2); + + for (;;) + { + if (!get_slots(winobj, winstate, relpos)) + { + out_of_frame = true; + break; /* current row is out of frame */ + } + + /* evaluate the expression */ + eval_result = ExecEvalExpr(pat, econtext, &isnull); + if (isnull) + { + /* expression is NULL */ + elog(DEBUG1, "expression for %s is NULL at row: %d", vname, relpos); + break; + } + else + { + if (!DatumGetBool(eval_result)) + { + /* expression is false */ + elog(DEBUG1, "expression for %s is false at row: %d", vname, relpos); + break; + } + else + { + /* expression is true */ + elog(DEBUG1, "expression for %s is true at row: %d", vname, relpos); + appendStringInfoChar(encoded_str, vname[0]); + + /* If quantifier is "+", we need to look for more matching row */ + if (quantifier && !strcmp(quantifier, "+")) + { + /* remember that we want to try another row */ + second_try_match = true; + relpos++; + } + else + break; + } + } + } + if (second_try_match) + relpos--; + + if (out_of_frame) + { + *result = false; + return -1; + } + + /* build regular expression */ + snprintf(pattern_str, sizeof(pattern_str), "%c%s", vname[0], quantifier); + + /* + * Do regular expression matching against sequence of rows satisfying + * the expression using regexp_instr(). + */ + sts = DatumGetInt32(DirectFunctionCall2Coll(regexp_instr, DEFAULT_COLLATION_OID, + PointerGetDatum(cstring_to_text(encoded_str->data)), + PointerGetDatum(cstring_to_text(pattern_str)))); + elog(DEBUG1, "regexp_instr returned: %d. str: %s regexp: %s", + sts, encoded_str->data, pattern_str); + *result = (sts > 0)? true : false; + } + return relpos; +} + +/* + * Get current, previous and next tuple. + * Returns true if still within frame. + */ +static bool +get_slots(WindowObject winobj, WindowAggState *winstate, int current_pos) +{ + TupleTableSlot *slot; + bool isnull, isout; + int sts; + ExprContext *econtext; + + econtext = winstate->ss.ps.ps_ExprContext; + + /* for current row */ + slot = winstate->temp_slot_1; + sts = WinGetSlotInFrame(winobj, slot, + current_pos, WINDOW_SEEK_HEAD, false, + &isnull, &isout); + if (sts < 0) + { + elog(DEBUG1, "current row is out of frame"); + econtext->ecxt_scantuple = winstate->null_slot; + return false; + } + else + econtext->ecxt_scantuple = slot; + + /* for PREV */ + if (current_pos > 0) + { + slot = winstate->prev_slot; + sts = WinGetSlotInFrame(winobj, slot, + current_pos - 1, WINDOW_SEEK_HEAD, false, + &isnull, &isout); + if (sts < 0) + { + elog(DEBUG1, "previous row out of frame at: %d", current_pos); + econtext->ecxt_outertuple = winstate->null_slot; + } + econtext->ecxt_outertuple = slot; + } + else + econtext->ecxt_outertuple = winstate->null_slot; + + /* for NEXT */ + slot = winstate->next_slot; + sts = WinGetSlotInFrame(winobj, slot, + current_pos + 1, WINDOW_SEEK_HEAD, false, + &isnull, &isout); + if (sts < 0) + { + elog(DEBUG1, "next row out of frame at: %d", current_pos); + econtext->ecxt_innertuple = winstate->null_slot; + } + else + econtext->ecxt_innertuple = slot; + + return true; +} + +/* + * prev + * Dummy function to invoke RPR's navigation operator "PREV". + * This is *not* a window function. + */ +Datum +window_prev(PG_FUNCTION_ARGS) +{ + PG_RETURN_DATUM(PG_GETARG_DATUM(0)); +} + +/* + * next + * Dummy function to invoke RPR's navigation operation "NEXT". + * This is *not* a window function. + */ +Datum +window_next(PG_FUNCTION_ARGS) +{ + PG_RETURN_DATUM(PG_GETARG_DATUM(0)); +} + diff --git a/src/include/catalog/pg_proc.dat b/src/include/catalog/pg_proc.dat index 6996073989..e3a9e0ffeb 100644 --- a/src/include/catalog/pg_proc.dat +++ b/src/include/catalog/pg_proc.dat @@ -10397,6 +10397,15 @@ { oid => '3114', descr => 'fetch the Nth row value', proname => 'nth_value', prokind => 'w', prorettype => 'anyelement', proargtypes => 'anyelement int4', prosrc => 'window_nth_value' }, +{ oid => '6122', descr => 'row pattern recognition in window', + proname => 'rpr', prokind => 'w', prorettype => 'anyelement', + proargtypes => 'anyelement', prosrc => 'window_rpr' }, +{ oid => '6123', descr => 'previous value', + proname => 'prev', provolatile => 's', prorettype => 'anyelement', + proargtypes => 'anyelement', prosrc => 'window_prev' }, +{ oid => '6124', descr => 'next value', + proname => 'next', provolatile => 's', prorettype => 'anyelement', + proargtypes => 'anyelement', prosrc => 'window_next' }, # functions for range types { oid => '3832', descr => 'I/O', diff --git a/src/include/nodes/execnodes.h b/src/include/nodes/execnodes.h index cb714f4a19..bc95c5fe9b 100644 --- a/src/include/nodes/execnodes.h +++ b/src/include/nodes/execnodes.h @@ -2519,6 +2519,13 @@ typedef struct WindowAggState int64 groupheadpos; /* current row's peer group head position */ int64 grouptailpos; /* " " " " tail position (group end+1) */ + /* these fields are used in Row pattern recognition: */ + List *patternVariableList; /* list of row pattern variables names (list of String) */ + List *patternRegexpList; /* list of row pattern regular expressions ('*', '+' or '?'. list of String) */ + List *defineVariableList; /* list of row pattern definition variables (list of String) */ + List *defineClauseList; /* expression for row pattern definition + * search conditions ExprState list */ + MemoryContext partcontext; /* context for partition-lifespan data */ MemoryContext aggcontext; /* shared context for aggregate working data */ MemoryContext curaggcontext; /* current aggregate's working data */ @@ -2555,6 +2562,11 @@ typedef struct WindowAggState TupleTableSlot *agg_row_slot; TupleTableSlot *temp_slot_1; TupleTableSlot *temp_slot_2; + + /* temporary slots for RPR */ + TupleTableSlot *prev_slot; /* PREV row navigation operator */ + TupleTableSlot *next_slot; /* NEXT row navigation operator */ + TupleTableSlot *null_slot; /* all NULL slot */ } WindowAggState; /* ---------------- diff --git a/src/include/windowapi.h b/src/include/windowapi.h index b8c2c565d1..a0facf38fe 100644 --- a/src/include/windowapi.h +++ b/src/include/windowapi.h @@ -58,7 +58,16 @@ extern Datum WinGetFuncArgInFrame(WindowObject winobj, int argno, int relpos, int seektype, bool set_mark, bool *isnull, bool *isout); +extern int WinGetSlotInFrame(WindowObject winobj, TupleTableSlot *slot, + int relpos, int seektype, bool set_mark, + bool *isnull, bool *isout); + extern Datum WinGetFuncArgCurrent(WindowObject winobj, int argno, bool *isnull); +extern WindowAggState *WinGetAggState(WindowObject winobj); + +extern bool window_gettupleslot(WindowObject winobj, int64 pos, TupleTableSlot *slot); + + #endif /* WINDOWAPI_H */ -- 2.25.1 ----Next_Part(Sun_Jun_25_21_05_09_2023_126)-- Content-Type: Text/X-Patch; charset=us-ascii Content-Transfer-Encoding: 7bit Content-Disposition: inline; filename="v1-0005-Row-pattern-recognition-patch-docs.patch" ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
* [PATCH] Lock upgrade without deadlocks. @ 2026-04-17 13:13 Antonin Houska <[email protected]> 0 siblings, 0 replies; 189+ messages in thread From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw) For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock to AccessExclusiveLock for the final stage of table processing. However, we cannot release the earlier before requesting the latter, because if another process ran DDL on the table in between, REPACK would have to cancel its transaction, and waste possibly a lot of work. Without releasing ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really deterministic whether REPACK or the other process will get canceled. This patch adds a new field to the LOCK structure which tells which process is trying to upgrade the lock - this field is set right before the upgrade. All processes waiting in the lock queue are then woken up to check the flag. When another process tries to get the lock after that, it checks this field, and if it's set, it checks for deadlocks even if deadlock timeout hasn't expired yet. If the process is already in the lock's queue and sleeping, the lock upgrading process wakes it up so it checks the flag immediately. At the time the upgrading process starts to sleep, all the other process should be aware that they need to check for deadlock during their lock acquisitions, so the upgrading process does not have to check for deadlocks when it gets woken up. Thus it should never receive deadlock error report. --- src/backend/commands/repack.c | 4 +- src/backend/storage/lmgr/lmgr.c | 35 ++++++++--- src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++- src/backend/storage/lmgr/proc.c | 32 +++++++--- src/include/storage/lmgr.h | 1 + src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 4 +- 7 files changed, 166 insertions(+), 22 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..310b2a65099 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there * is one), all its indexes, so that we can swap the files. + * + * TODO The same for indexes and TOAST? */ - LockRelationOid(old_table_oid, AccessExclusiveLock); + LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock); /* * Lock all indexes now, not only the clustering one: all indexes need to diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c index 2ccf7237fee..0b103485d54 100644 --- a/src/backend/storage/lmgr/lmgr.c +++ b/src/backend/storage/lmgr/lmgr.c @@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo const ItemPointerData *ctid; } XactLockTableWaitInfo; +static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode, + bool isUpgrade); static void XactLockTableWaitErrorCb(void *arg); /* @@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid) */ void LockRelationOid(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, false); +} + +/* + * Like above, but upgrade an existing lock. + */ +void +LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode) +{ + LockRelationOidCommon(relid, lockmode, true); +} + +static void +LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade) { LOCKTAG tag; LOCALLOCK *locallock; @@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, isUpgrade); /* * Now that we have the lock, check for invalidation messages, so that we @@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode) SetLocktagRelationOid(&tag, relid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode) SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock, - false); + false, false); /* * Now that we have the lock, check for invalidation messages; see notes @@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode) relation->rd_lockInfo.lockRelId.relId); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc ItemPointerGetOffsetNumber(tid)); return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL, - logLockFailure) != LOCKACQUIRE_NOT_AVAIL); + logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL); } /* @@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure) SET_LOCKTAG_TRANSACTION(tag, xid); if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL, - logLockFailure) + logLockFailure, false) == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; @@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid, objsubid); res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock, - false); + false, false); if (res == LOCKACQUIRE_NOT_AVAIL) return false; diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..bcbbf36c091 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag, bool dontWait) { return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait, - true, NULL, false); + true, NULL, false, false); } /* @@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag, * * logLockFailure indicates whether to log details when a lock acquisition * fails with dontWait = true. + * + * isUpgrade should be true if the backend already holds this lock in a lower + * mode. */ LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, @@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure) + bool logLockFailure, + bool isUpgrade) { LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid; LockMethod lockMethodTable; @@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag, * case, because JoinWaitQueue() may discover that we can acquire the * lock immediately after all. */ - waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait); + waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait, + isUpgrade); } if (waitResult == PROC_WAIT_STATUS_ERROR) @@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag, Assert(!dontWait); PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock); LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode); + + /* + * Lock upgrade can introduce deadlock, therefore enforce special + * behavior of other processes that deal with this lock. In + * particular, any waiter that sees upgradedBy set is expected to + * perform deadlock check as soon as it's woken up. (Since we are + * already in the queue, the deadlock detector has all the information + * it needs.) + */ + if (isUpgrade) + { + dlist_iter iter; + + /* + * There should not be multiple upgrades at the same time. XXX + * ERROR ? + */ + Assert(lock->upgradedBy == NULL); + + lock->upgradedBy = MyProc; + + /* + * Now, before I sleep myself, wake up all the existing waiters + * (except for me), so they check for deadlock. + */ + dclist_foreach(iter, &lock->waitProcs) + { + PGPROC *proc = dlist_container(PGPROC, waitLink, + iter.cur); + + if (proc != MyProc) + SetLatch(&proc->procLatch); + } + } LWLockRelease(partitionLock); + if (!isUpgrade) + { + HASH_SEQ_STATUS status; + LOCALLOCK *llock; + bool check_deadlock = false; + + /* + * Likewise, new waiters should check the isUpgrade flag and + * engage deadlock detector if needed. The lock being acquired + * should already be in the hash. + * + * Note that, to make deadlock detection happen as soon as + * possible (i.e. regardless deadlock timeout), we check the other + * locks too. XXX There might be ways to get into the deadlock + * indirectly, but that needs more analysis. As long as the + * upgrading process skips deadlock detection altogether, the + * worst consequence of missing some lock wait cycle here is that + * the other process will be kicked-off no sooner than after its + * deadlock timeout has elapsed. (Which in turn means that the + * lock upgrade will take longer than expected.) + */ + hash_seq_init(&status, LockMethodLocalHash); + while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL) + { + LOCK *mylock = llock->lock; + + if (mylock && mylock->upgradedBy) + { + check_deadlock = true; + break; + } + } + /* Check for deadlock if needed. */ + if (check_deadlock) + { + DeadLockState deadlock_state; + + deadlock_state = CheckDeadLock(); + if (deadlock_state == DS_HARD_DEADLOCK) + DeadLockReport(); + } + } + waitResult = WaitOnLock(locallock, owner); /* @@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag, DeadLockReport(); /* DeadLockReport() will not return */ } + + /* + * If finishing the lock upgrade, we should not get into a deadlock + * anymore, so let others know that they do not have to care either. + */ + if (isUpgrade) + { + LWLockAcquire(partitionLock, LW_EXCLUSIVE); + Assert(lock->upgradedBy != NULL); + lock->upgradedBy = NULL; + LWLockRelease(partitionLock); + } } else LWLockRelease(partitionLock); @@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc, lock->nGranted = 0; MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES); MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES); + lock->upgradedBy = NULL; LOCK_PRINT("LockAcquire: new", lock, lockmode); } else @@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock, elog(PANIC, "proclock table corrupted"); } + /* + * Was this backend upgrading the lock? + */ + if (lock->upgradedBy == MyProc) + lock->upgradedBy = NULL; + if (lock->nRequested == 0) { /* diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..7b31370db70 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout; static void RemoveProcFromArray(int code, Datum arg); static void ProcKill(int code, Datum arg); static void AuxiliaryProcKill(int code, Datum arg); -static DeadLockState CheckDeadLock(void); /* @@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid) * NOTES: The process queue is now a priority queue for locking. */ ProcWaitStatus -JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) +JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait, + bool isUpgrade) { LOCKMODE lockmode = locallock->tag.mode; LOCK *lock = locallock->lock; @@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) /* Must he wait for me? */ if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks) { - /* Must I wait for him ? */ - if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks) + /* + * Must I wait for him? I don't want a deadlock during lock + * upgrade - other processes should fail on it. + */ + if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) && + !isUpgrade) { /* * Yes, so we have a deadlock. Easiest way to clean up @@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock) (void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0, PG_WAIT_LOCK | locallock->tag.lock.locktag_type); ResetLatch(MyLatch); - /* check for deadlocks first, as that's probably log-worthy */ - if (got_deadlock_timeout) + /* + * Check for deadlocks first, as that's probably log-worthy. + * + * Do not wait for the timeout if the lock is being upgraded since + * the risk of deadlock is higher now. However, do not check for + * deadlock if the lock is being upgraded by this process - other + * processes should take care. + * + * TODO Possible optimization: if this is the only lock of the + * backend and if it did not have any weaker lock on the table so + * far, it should be safe to skip the deadlock check. However, to + * evaluate the situation, we need to take fast-path locks into + * account. + */ + if ((got_deadlock_timeout || lock->upgradedBy) && + lock->upgradedBy != MyProc) { deadlock_state = CheckDeadLock(); got_deadlock_timeout = false; @@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock) * not, just return. If we have a real deadlock, remove ourselves from the * lock's wait queue. */ -static DeadLockState +DeadLockState CheckDeadLock(void) { int i; diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h index 2a985ce5e15..eddbe1e8622 100644 --- a/src/include/storage/lmgr.h +++ b/src/include/storage/lmgr.h @@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation); /* Lock a relation */ extern void LockRelationOid(Oid relid, LOCKMODE lockmode); +extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode); extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode); extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode); extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode); diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..080b69eddfd 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod; * nRequested -- total requested locks of all types. * granted -- count of each lock type currently granted on the lock. * nGranted -- total granted locks of all types. + * upgrader -- process that intends to upgrade the lock * * Note: these counts count 1 for each backend. Internally to a backend, * there may be multiple grabs on a particular lock, but this is not reflected @@ -150,6 +151,7 @@ typedef struct LOCK int nRequested; /* total of requested[] array */ int granted[MAX_LOCKMODES]; /* counts of granted locks */ int nGranted; /* total of granted[] array */ + PGPROC *upgradedBy; /* is lock being upgraded by this process? */ } LOCK; #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid) @@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag, bool dontWait, bool reportMemoryError, LOCALLOCK **locallockp, - bool logLockFailure); + bool logLockFailure, + bool isUpgrade); extern void AbortStrongLockAcquire(void); extern void MarkLockClear(LOCALLOCK *locallock); extern bool LockRelease(const LOCKTAG *locktag, diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..c4bc3a24044 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree); extern void ProcReleaseLocks(bool isCommit); extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock, - LockMethod lockMethodTable, bool dontWait); + LockMethod lockMethodTable, bool dontWait, + bool isUpgrade); extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock); extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus); extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock); +extern DeadLockState CheckDeadLock(void); extern void CheckDeadLockAlert(void); extern void LockErrorCleanup(void); extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock, -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 189+ messages in thread
end of thread, other threads:[~2026-04-17 13:13 UTC | newest] Thread overview: 189+ messages (download: mbox mbox.gz follow: Atom feed) -- links below jump to the message on this page -- 2023-06-25 11:48 [PATCH v1 4/7] Row pattern recognition patch (executor). Tatsuo Ishii <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]> 2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
This inbox is served by agora; see mirroring instructions for how to clone and mirror all data and code used for this inbox